Micron Document
<!DOCTYPE html>
<html class="client-nojs vector-feature-night-mode-disabled vector-feature-language-in-header-enabled vector-feature-language-in-main-page-header-disabled vector-feature-page-tools-pinned-disabled vector-feature-toc-pinned-clientpref-1 vector-feature-main-menu-pinned-disabled vector-feature-limited-width-clientpref-1 vector-feature-limited-width-content-enabled vector-feature-custom-font-size-clientpref-1 vector-feature-appearance-pinned-clientpref-1 vector-sticky-header-enabled" lang="en" dir="ltr"><head>
<meta charset="UTF-8">
<title>Multimodal learning</title>
<meta name="viewport" content="width=device-width, initial-scale=1.0">
<link rel="canonical" href="https://en.wikipedia.org/wiki/Multimodal_learning"> <link href="./mw/ext.cite.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/ext.math.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.icons.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.search.codex.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/skins.vector.styles.css" rel="stylesheet" type="text/css">
<link href="./mw/user.styles.css" rel="stylesheet" type="text/css">
<meta name="ResourceLoaderDynamicStyles" content="">
<link rel="stylesheet" type="text/css" href="./mw/site.styles.css">
<link rel="stylesheet" type="text/css" href="./mw/noscript.css">
<link rel="stylesheet" type="text/css" href="./footer.css">
<link rel="stylesheet" type="text/css" href="./vector-2022.css">
</head>
<body class="skin--responsive skin-vector skin-vector-search-vue mediawiki ltr sitedir-ltr mw-hide-empty-elt ns-0 ns-subject page-Multimodal_learning rootpage-Multimodal_learning skin-vector-2022 action-view">
<div class="mw-page-container">
<div class="mw-page-container-inner">
<div class="mw-content-container">
<main id="content" class="mw-body">
<header class="mw-body-header vector-page-titlebar">
<h1 id="firstHeading" class="firstHeading mw-first-heading">
<span id="openzim-page-title" class="mw-page-title-main"><span class="mw-page-title-main">Multimodal learning</span></span>
</h1>
</header>
<a id="top"></a>
<div id="bodyContent" class="vector-body ve-init-mw-desktopArticleTarget-targetContainer" aria-labelledby="firstHeading" data-mw-ve-target-container="">
<div id="mw-content-text" class="mw-body-content mw-content-ltr" lang="en" dir="ltr"><div class="mw-content-ltr mw-parser-output" lang="en" dir="ltr">
<style data-mw-deduplicate="TemplateStyles:r1129693374">
/* start https://en.wikipedia.org/ */


.mw-parser-output .hlist dl,.mw-parser-output .hlist ol,.mw-parser-output .hlist ul{margin:0;padding:0}.mw-parser-output .hlist dd,.mw-parser-output .hlist dt,.mw-parser-output .hlist li{margin:0;display:inline}.mw-parser-output .hlist.inline,.mw-parser-output .hlist.inline dl,.mw-parser-output .hlist.inline ol,.mw-parser-output .hlist.inline ul,.mw-parser-output .hlist dl dl,.mw-parser-output .hlist dl ol,.mw-parser-output .hlist dl ul,.mw-parser-output .hlist ol dl,.mw-parser-output .hlist ol ol,.mw-parser-output .hlist ol ul,.mw-parser-output .hlist ul dl,.mw-parser-output .hlist ul ol,.mw-parser-output .hlist ul ul{display:inline}.mw-parser-output .hlist .mw-empty-li{display:none}.mw-parser-output .hlist dt::after{content:": "}.mw-parser-output .hlist dd::after,.mw-parser-output .hlist li::after{content:" · ";font-weight:bold}.mw-parser-output .hlist dd:last-child::after,.mw-parser-output .hlist dt:last-child::after,.mw-parser-output .hlist li:last-child::after{content:none}.mw-parser-output .hlist dd dd:first-child::before,.mw-parser-output .hlist dd dt:first-child::before,.mw-parser-output .hlist dd li:first-child::before,.mw-parser-output .hlist dt dd:first-child::before,.mw-parser-output .hlist dt dt:first-child::before,.mw-parser-output .hlist dt li:first-child::before,.mw-parser-output .hlist li dd:first-child::before,.mw-parser-output .hlist li dt:first-child::before,.mw-parser-output .hlist li li:first-child::before{content:" (";font-weight:normal}.mw-parser-output .hlist dd dd:last-child::after,.mw-parser-output .hlist dd dt:last-child::after,.mw-parser-output .hlist dd li:last-child::after,.mw-parser-output .hlist dt dd:last-child::after,.mw-parser-output .hlist dt dt:last-child::after,.mw-parser-output .hlist dt li:last-child::after,.mw-parser-output .hlist li dd:last-child::after,.mw-parser-output .hlist li dt:last-child::after,.mw-parser-output .hlist li li:last-child::after{content:")";font-weight:normal}.mw-parser-output .hlist ol{counter-reset:listitem}.mw-parser-output .hlist ol>li{counter-increment:listitem}.mw-parser-output .hlist ol>li::before{content:" "counter(listitem)"\a0 "}.mw-parser-output .hlist dd ol>li:first-child::before,.mw-parser-output .hlist dt ol>li:first-child::before,.mw-parser-output .hlist li ol>li:first-child::before{content:" ("counter(listitem)"\a0 "}


/* end https://en.wikipedia.org/ */
</style><style data-mw-deduplicate="TemplateStyles:r1246091330">
/* start https://en.wikipedia.org/ */


.mw-parser-output .sidebar{width:22em;float:right;clear:right;margin:0.5em 0 1em 1em;background:var(--background-color-neutral-subtle,#f8f9fa);border:1px solid var(--border-color-base,#a2a9b1);padding:0.2em;text-align:center;line-height:1.4em;font-size:88%;border-collapse:collapse;display:table}body.skin-minerva .mw-parser-output .sidebar{display:table!important;float:right!important;margin:0.5em 0 1em 1em!important}.mw-parser-output .sidebar-subgroup{width:100%;margin:0;border-spacing:0}.mw-parser-output .sidebar-left{float:left;clear:left;margin:0.5em 1em 1em 0}.mw-parser-output .sidebar-none{float:none;clear:both;margin:0.5em 1em 1em 0}.mw-parser-output .sidebar-outer-title{padding:0 0.4em 0.2em;font-size:125%;line-height:1.2em;font-weight:bold}.mw-parser-output .sidebar-top-image{padding:0.4em}.mw-parser-output .sidebar-top-caption,.mw-parser-output .sidebar-pretitle-with-top-image,.mw-parser-output .sidebar-caption{padding:0.2em 0.4em 0;line-height:1.2em}.mw-parser-output .sidebar-pretitle{padding:0.4em 0.4em 0;line-height:1.2em}.mw-parser-output .sidebar-title,.mw-parser-output .sidebar-title-with-pretitle{padding:0.2em 0.8em;font-size:145%;line-height:1.2em}.mw-parser-output .sidebar-title-with-pretitle{padding:0.1em 0.4em}.mw-parser-output .sidebar-image{padding:0.2em 0.4em 0.4em}.mw-parser-output .sidebar-heading{padding:0.1em 0.4em}.mw-parser-output .sidebar-content{padding:0 0.5em 0.4em}.mw-parser-output .sidebar-content-with-subgroup{padding:0.1em 0.4em 0.2em}.mw-parser-output .sidebar-above,.mw-parser-output .sidebar-below{padding:0.3em 0.8em;font-weight:bold}.mw-parser-output .sidebar-collapse .sidebar-above,.mw-parser-output .sidebar-collapse .sidebar-below{border-top:1px solid #aaa;border-bottom:1px solid #aaa}.mw-parser-output .sidebar-navbar{text-align:right;font-size:115%;padding:0 0.4em 0.4em}.mw-parser-output .sidebar-list-title{padding:0 0.4em;text-align:left;font-weight:bold;line-height:1.6em;font-size:105%}.mw-parser-output .sidebar-list-title-c{padding:0 0.4em;text-align:center;margin:0 3.3em}@media(max-width:640px){body.mediawiki .mw-parser-output .sidebar{width:100%!important;clear:both;float:none!important;margin-left:0!important;margin-right:0!important}}body.skin--responsive .mw-parser-output .sidebar a>img{max-width:none!important}@media screen{html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-list-title,html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle{background:transparent!important}html.skin-theme-clientpref-night .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle a{color:var(--color-progressive)!important}}@media screen and (prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-list-title,html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle{background:transparent!important}html.skin-theme-clientpref-os .mw-parser-output .sidebar:not(.notheme) .sidebar-title-with-pretitle a{color:var(--color-progressive)!important}}@media print{body.ns-0 .mw-parser-output .sidebar{display:none!important}}


/* end https://en.wikipedia.org/ */
</style><style data-mw-deduplicate="TemplateStyles:r886047488">
/* start https://en.wikipedia.org/ */


.mw-parser-output .nobold{font-weight:normal}


/* end https://en.wikipedia.org/ */
</style><table class="sidebar sidebar-collapse nomobile nowraplinks"><tbody><tr><td class="sidebar-pretitle">Part of a series on</td></tr><tr><th class="sidebar-title-with-pretitle"><a href="Machine_learning" title="Machine learning">Machine learning</a><br>and <a href="Data_mining" title="Data mining">data mining</a></th></tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Paradigms</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Supervised_learning" title="Supervised learning">Supervised learning</a></li>
<li><a href="Unsupervised_learning" title="Unsupervised learning">Unsupervised learning</a></li>
<li><a href="Semi-supervised_learning" class="mw-redirect" title="Semi-supervised learning">Semi-supervised learning</a></li>
<li><a href="Self-supervised_learning" title="Self-supervised learning">Self-supervised learning</a></li>
<li><a href="Reinforcement_learning" title="Reinforcement learning">Reinforcement learning</a></li>
<li><a href="Meta-learning_(computer_science)" title="Meta-learning (computer science)">Meta-learning</a></li>
<li><a href="Online_machine_learning" title="Online machine learning">Online learning</a></li>
<li><a href="Batch_learning" class="mw-redirect" title="Batch learning">Batch learning</a></li>
<li><a href="Curriculum_learning" title="Curriculum learning">Curriculum learning</a></li>
<li><a href="Rule-based_machine_learning" title="Rule-based machine learning">Rule-based learning</a></li>
<li><a href="Neuro-symbolic_AI" title="Neuro-symbolic AI">Neuro-symbolic AI</a></li>
<li><a href="Neuromorphic_engineering" class="mw-redirect" title="Neuromorphic engineering">Neuromorphic engineering</a></li>
<li><a href="Quantum_machine_learning" title="Quantum machine learning">Quantum machine learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Problems</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Statistical_classification" title="Statistical classification">Classification</a></li>
<li><a href="Generative_model" title="Generative model">Generative modeling</a></li>
<li><a href="Regression_analysis" title="Regression analysis">Regression</a></li>
<li><a href="Cluster_analysis" title="Cluster analysis">Clustering</a></li>
<li><a href="Dimensionality_reduction" title="Dimensionality reduction">Dimensionality reduction</a></li>
<li><a href="Density_estimation" title="Density estimation">Density estimation</a></li>
<li><a href="Anomaly_detection" title="Anomaly detection">Anomaly detection</a></li>
<li><a href="Data_cleaning" class="mw-redirect" title="Data cleaning">Data cleaning</a></li>
<li><a href="Automated_machine_learning" title="Automated machine learning">AutoML</a></li>
<li><a href="Association_rule_learning" title="Association rule learning">Association rules</a></li>
<li><a href="Semantic_analysis_(machine_learning)" title="Semantic analysis (machine learning)">Semantic analysis</a></li>
<li><a href="Structured_prediction" title="Structured prediction">Structured prediction</a></li>
<li><a href="Feature_engineering" title="Feature engineering">Feature engineering</a></li>
<li><a href="Feature_learning" title="Feature learning">Feature learning</a></li>
<li><a href="Learning_to_rank" title="Learning to rank">Learning to rank</a></li>
<li><a href="Grammar_induction" title="Grammar induction">Grammar induction</a></li>
<li><a href="Ontology_learning" title="Ontology learning">Ontology learning</a></li>
</ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><div style="display: inline-block; line-height: 1.2em; padding: .1em 0;"><a href="Supervised_learning" title="Supervised learning">Supervised learning</a><br><span class="nobold"><span style="font-size: 85%;">(<b><a href="Statistical_classification" title="Statistical classification">classification</a></b>&nbsp;• <b><a href="Regression_analysis" title="Regression analysis">regression</a></b>)</span></span> </div></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Apprenticeship_learning" title="Apprenticeship learning">Apprenticeship learning</a></li>
<li><a href="Decision_tree_learning" title="Decision tree learning">Decision trees</a></li>
<li><a href="Ensemble_learning" title="Ensemble learning">Ensembles</a>
<ul><li><a href="Bootstrap_aggregating" title="Bootstrap aggregating">Bagging</a></li>
<li><a href="Boosting_(machine_learning)" title="Boosting (machine learning)">Boosting</a></li>
<li><a href="Random_forest" title="Random forest">Random forest</a></li></ul></li>
<li><a href="K-nearest_neighbors_algorithm" title="K-nearest neighbors algorithm"><i>k</i>-NN</a></li>
<li><a href="Linear_regression" title="Linear regression">Linear regression</a></li>
<li><a href="Naive_Bayes_classifier" title="Naive Bayes classifier">Naive Bayes</a></li>
<li><a href="Artificial_neural_network" class="mw-redirect" title="Artificial neural network">Artificial neural networks</a></li>
<li><a href="Logistic_regression" title="Logistic regression">Logistic regression</a></li>
<li><a href="Perceptron" title="Perceptron">Perceptron</a></li>
<li><a href="Relevance_vector_machine" title="Relevance vector machine">Relevance vector machine (RVM)</a></li>
<li><a href="Support_vector_machine" title="Support vector machine">Support vector machine (SVM)</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Cluster_analysis" title="Cluster analysis">Clustering</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="BIRCH" title="BIRCH">BIRCH</a></li>
<li><a href="CURE_algorithm" title="CURE algorithm">CURE</a></li>
<li><a href="Hierarchical_clustering" title="Hierarchical clustering">Hierarchical</a></li>
<li><a href="K-means_clustering" title="K-means clustering"><i>k</i>-means</a></li>
<li><a href="Fuzzy_clustering" title="Fuzzy clustering">Fuzzy</a></li>
<li><a href="Expectation%E2%80%93maximization_algorithm" title="Expectation–maximization algorithm">Expectation–maximization (EM)</a></li>
<li><br><a href="DBSCAN" title="DBSCAN">DBSCAN</a></li>
<li><a href="OPTICS_algorithm" title="OPTICS algorithm">OPTICS</a></li>
<li><a href="Mean_shift" title="Mean shift">Mean shift</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Dimensionality_reduction" title="Dimensionality reduction">Dimensionality reduction</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Factor_analysis" title="Factor analysis">Factor analysis</a></li>
<li><a href="Canonical_correlation" title="Canonical correlation">CCA</a></li>
<li><a href="Independent_component_analysis" title="Independent component analysis">ICA</a></li>
<li><a href="Linear_discriminant_analysis" title="Linear discriminant analysis">LDA</a></li>
<li><a href="Non-negative_matrix_factorization" title="Non-negative matrix factorization">NMF</a></li>
<li><a href="Principal_component_analysis" title="Principal component analysis">PCA</a></li>
<li><a href="Proper_generalized_decomposition" title="Proper generalized decomposition">PGD</a></li>
<li><a href="T-distributed_stochastic_neighbor_embedding" title="T-distributed stochastic neighbor embedding">t-SNE</a></li>
<li><a href="Sparse_dictionary_learning" title="Sparse dictionary learning">SDL</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Structured_prediction" title="Structured prediction">Structured prediction</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Graphical_model" title="Graphical model">Graphical models</a>
<ul><li><a href="Bayesian_network" title="Bayesian network">Bayes net</a></li>
<li><a href="Conditional_random_field" title="Conditional random field">Conditional random field</a></li>
<li><a href="Hidden_Markov_model" title="Hidden Markov model">Hidden Markov</a></li></ul></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Anomaly_detection" title="Anomaly detection">Anomaly detection</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Random_sample_consensus" title="Random sample consensus">RANSAC</a></li>
<li><a href="K-nearest_neighbors_algorithm" title="K-nearest neighbors algorithm"><i>k</i>-NN</a></li>
<li><a href="Local_outlier_factor" title="Local outlier factor">Local outlier factor</a></li>
<li><a href="Isolation_forest" title="Isolation forest">Isolation forest</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Neural_network_(machine_learning)" title="Neural network (machine learning)">Neural networks</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Autoencoder" title="Autoencoder">Autoencoder</a></li>
<li><a href="Deep_learning" title="Deep learning">Deep learning</a></li>
<li><a href="Feedforward_neural_network" title="Feedforward neural network">Feedforward neural network</a></li>
<li><a href="Recurrent_neural_network" title="Recurrent neural network">Recurrent neural network</a>
<ul><li><a href="Long_short-term_memory" title="Long short-term memory">LSTM</a></li>
<li><a href="Gated_recurrent_unit" title="Gated recurrent unit">GRU</a></li>
<li><a href="Echo_state_network" title="Echo state network">ESN</a></li>
<li><a href="Reservoir_computing" title="Reservoir computing">reservoir computing</a></li></ul></li>
<li><a href="Boltzmann_machine" title="Boltzmann machine">Boltzmann machine</a>
<ul><li><a href="Restricted_Boltzmann_machine" title="Restricted Boltzmann machine">Restricted</a></li></ul></li>
<li><a href="Generative_adversarial_network" title="Generative adversarial network">GAN</a></li>
<li><a href="Diffusion_model" title="Diffusion model">Diffusion model</a></li>
<li><a href="Self-organizing_map" title="Self-organizing map">SOM</a></li>
<li><a href="Convolutional_neural_network" title="Convolutional neural network">Convolutional neural network</a>
<ul><li><a href="U-Net" title="U-Net">U-Net</a></li>
<li><a href="LeNet" title="LeNet">LeNet</a></li>
<li><a href="AlexNet" title="AlexNet">AlexNet</a></li>
<li><a href="DeepDream" title="DeepDream">DeepDream</a></li></ul></li>
<li><a href="Neural_field" title="Neural field">Neural field</a>
<ul><li><a href="Neural_radiance_field" title="Neural radiance field">Neural radiance field</a></li>
<li><a href="Physics-informed_neural_networks" title="Physics-informed neural networks">Physics-informed neural networks</a></li></ul></li>
<li><a href="Transformer_(deep_learning_architecture)" title="Transformer (deep learning architecture)">Transformer</a>
<ul><li><a href="Vision_transformer" title="Vision transformer">Vision</a></li></ul></li>
<li><a href="Mamba_(deep_learning_architecture)" title="Mamba (deep learning architecture)">Mamba</a></li>
<li><a href="Spiking_neural_network" title="Spiking neural network">Spiking neural network</a></li>
<li><a href="Memtransistor" title="Memtransistor">Memtransistor</a></li>
<li><a href="Electrochemical_RAM" title="Electrochemical RAM">Electrochemical RAM</a> (ECRAM)</li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)"><a href="Reinforcement_learning" title="Reinforcement learning">Reinforcement learning</a></div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Q-learning" title="Q-learning">Q-learning</a></li>
<li><a href="Policy_gradient_method" title="Policy gradient method">Policy gradient</a></li>
<li><a href="State%E2%80%93action%E2%80%93reward%E2%80%93state%E2%80%93action" title="State–action–reward–state–action">SARSA</a></li>
<li><a href="Temporal_difference_learning" title="Temporal difference learning">Temporal difference (TD)</a></li>
<li><a href="Multi-agent_reinforcement_learning" title="Multi-agent reinforcement learning">Multi-agent</a>
<ul><li><a href="Self-play_(reinforcement_learning_technique)" class="mw-redirect" title="Self-play (reinforcement learning technique)">Self-play</a></li></ul></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Learning with humans</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Active_learning_(machine_learning)" title="Active learning (machine learning)">Active learning</a></li>
<li><a href="Crowdsourcing" title="Crowdsourcing">Crowdsourcing</a></li>
<li><a href="Human-in-the-loop" title="Human-in-the-loop">Human-in-the-loop</a></li>
<li><a href="Mechanistic_interpretability" title="Mechanistic interpretability">Mechanistic interpretability</a></li>
<li><a href="Reinforcement_learning_from_human_feedback" title="Reinforcement learning from human feedback">RLHF</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Model diagnostics</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Coefficient_of_determination" title="Coefficient of determination">Coefficient of determination</a></li>
<li><a href="Confusion_matrix" title="Confusion matrix">Confusion matrix</a></li>
<li><a href="Learning_curve_(machine_learning)" title="Learning curve (machine learning)">Learning curve</a></li>
<li><a href="Receiver_operating_characteristic" title="Receiver operating characteristic">ROC curve</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Mathematical foundations</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Kernel_machines" class="mw-redirect" title="Kernel machines">Kernel machines</a></li>
<li><a href="Bias%E2%80%93variance_tradeoff" title="Bias–variance tradeoff">Bias–variance tradeoff</a></li>
<li><a href="Computational_learning_theory" title="Computational learning theory">Computational learning theory</a></li>
<li><a href="Empirical_risk_minimization" title="Empirical risk minimization">Empirical risk minimization</a></li>
<li><a href="Occam_learning" title="Occam learning">Occam learning</a></li>
<li><a href="Probably_approximately_correct_learning" title="Probably approximately correct learning">PAC learning</a></li>
<li><a href="Statistical_learning_theory" title="Statistical learning theory">Statistical learning</a></li>
<li><a href="Vapnik%E2%80%93Chervonenkis_theory" title="Vapnik–Chervonenkis theory">VC theory</a></li>
<li><a href="Topological_deep_learning" title="Topological deep learning">Topological deep learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Journals and conferences</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="AAAI_Conference_on_Artificial_Intelligence" title="AAAI Conference on Artificial Intelligence">AAAI</a></li>
<li><a href="ECML_PKDD" title="ECML PKDD">ECML PKDD</a></li>
<li><a href="Conference_on_Neural_Information_Processing_Systems" title="Conference on Neural Information Processing Systems">NeurIPS</a></li>
<li><a href="International_Conference_on_Machine_Learning" title="International Conference on Machine Learning">ICML</a></li>
<li><a href="International_Conference_on_Learning_Representations" title="International Conference on Learning Representations">ICLR</a></li>
<li><a href="International_Joint_Conference_on_Artificial_Intelligence" title="International Joint Conference on Artificial Intelligence">IJCAI</a></li>
<li><a href="Machine_Learning_(journal)" title="Machine Learning (journal)">ML</a></li>
<li><a href="Journal_of_Machine_Learning_Research" title="Journal of Machine Learning Research">JMLR</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-content">
<div class="sidebar-list mw-collapsible mw-collapsed machine-learning-list-title"><div class="sidebar-list-title" style="border-top:1px solid #aaa; text-align:center;;color: var(--color-base)">Related articles</div><div class="sidebar-list-content mw-collapsible-content hlist">
<ul><li><a href="Glossary_of_artificial_intelligence" title="Glossary of artificial intelligence">Glossary of artificial intelligence</a></li>
<li><a href="List_of_datasets_for_machine-learning_research" title="List of datasets for machine-learning research">List of datasets for machine-learning research</a>
<ul><li><a href="List_of_datasets_in_computer_vision_and_image_processing" title="List of datasets in computer vision and image processing">List of datasets in computer vision and image processing</a></li></ul></li>
<li><a href="Outline_of_machine_learning" title="Outline of machine learning">Outline of machine learning</a></li></ul></div></div></td>
</tr><tr><td class="sidebar-navbar"><style data-mw-deduplicate="TemplateStyles:r1239400231">
/* start https://en.wikipedia.org/ */


.mw-parser-output .navbar{display:inline;font-size:88%;font-weight:normal}.mw-parser-output .navbar-collapse{float:left;text-align:left}.mw-parser-output .navbar-boxtext{word-spacing:0}.mw-parser-output .navbar ul{display:inline-block;white-space:nowrap;line-height:inherit}.mw-parser-output .navbar-brackets::before{margin-right:-0.125em;content:"[ "}.mw-parser-output .navbar-brackets::after{margin-left:-0.125em;content:" ]"}.mw-parser-output .navbar li{word-spacing:-0.125em}.mw-parser-output .navbar a>span,.mw-parser-output .navbar a>abbr{text-decoration:inherit}.mw-parser-output .navbar-mini abbr{font-variant:small-caps;border-bottom:none;text-decoration:none;cursor:inherit}.mw-parser-output .navbar-ct-full{font-size:114%;margin:0 7em}.mw-parser-output .navbar-ct-mini{font-size:114%;margin:0 4em}html.skin-theme-clientpref-night .mw-parser-output .navbar li a abbr{color:var(--color-base)!important}@media(prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .navbar li a abbr{color:var(--color-base)!important}}@media print{.mw-parser-output .navbar{display:none!important}}


/* end https://en.wikipedia.org/ */
</style></td></tr></tbody></table>
<p><b>Multimodal learning</b> is a type of <a href="Deep_learning" title="Deep learning">deep learning</a> that integrates and processes multiple types of data, referred to as <a href="Modality_(human%E2%80%93computer_interaction)" title="Modality (human–computer interaction)">modalities</a>, such as text, audio, images, or video. This integration allows for a more holistic understanding of complex data, improving model performance in tasks like visual question answering, cross-modal retrieval,<sup id="cite_ref-1" class="reference"><a href="#cite_note-1"><span class="cite-bracket">[</span>1<span class="cite-bracket">]</span></a></sup> text-to-image generation,<sup id="cite_ref-stable-diffusion-github_2-0" class="reference"><a href="#cite_note-stable-diffusion-github-2"><span class="cite-bracket">[</span>2<span class="cite-bracket">]</span></a></sup> aesthetic ranking,<sup id="cite_ref-3" class="reference"><a href="#cite_note-3"><span class="cite-bracket">[</span>3<span class="cite-bracket">]</span></a></sup> and image captioning.<sup id="cite_ref-4" class="reference"><a href="#cite_note-4"><span class="cite-bracket">[</span>4<span class="cite-bracket">]</span></a></sup>
</p><p>Large multimodal models, such as <a href="Google_Gemini" class="mw-redirect" title="Google Gemini">Google Gemini</a> and <a href="GPT-4o" title="GPT-4o">GPT-4o</a>, have become increasingly popular since 2023, enabling increased versatility and a broader understanding of real-world phenomena.<sup id="cite_ref-5" class="reference"><a href="#cite_note-5"><span class="cite-bracket">[</span>5<span class="cite-bracket">]</span></a></sup>
</p>
<meta property="mw:PageProp/toc">
<div class="mw-heading mw-heading2"><h2 id="Motivation">Motivation</h2></div>
<p>Data usually comes with different modalities which carry different information. For example, it is very common to caption an image to convey the information not presented in the image itself. Similarly, sometimes it is more straightforward to use an image to describe information which may not be obvious from text. As a result, if different words appear in similar images, then these words likely describe the same thing. Conversely, if a word is used to describe seemingly dissimilar images, then these images may represent the same object. Thus, in cases dealing with multi-modal data, it is important to use a model which is able to jointly represent the information such that the model can capture the combined information from different modalities.
</p>
<div class="mw-heading mw-heading2"><h2 id="Multimodal_transformers">Multimodal transformers</h2></div>
<div class="excerpt-block"><style data-mw-deduplicate="TemplateStyles:r1066933788">
/* start https://en.wikipedia.org/ */


.mw-parser-output .excerpt-hat .mw-editsection-like{font-style:normal}


/* end https://en.wikipedia.org/ */
</style><style data-mw-deduplicate="TemplateStyles:r1236090951">
/* start https://en.wikipedia.org/ */


.mw-parser-output .hatnote{font-style:italic}.mw-parser-output div.hatnote{padding-left:1.6em;margin-bottom:0.5em}.mw-parser-output .hatnote i{font-style:normal}.mw-parser-output .hatnote+link+.hatnote{margin-top:-0.5em}@media print{body.ns-0 .mw-parser-output .hatnote{display:none!important}}


/* end https://en.wikipedia.org/ */
</style><div role="note" class="hatnote navigation-not-searchable dablink excerpt-hat selfref">This section is an excerpt from <a href="Transformer_(deep_learning_architecture)#Multimodality" title="Transformer (deep learning architecture)">Transformer (deep learning architecture) § Multimodality</a>.<span class="mw-editsection-like "><span class="mw-editsection-bracket">[</span><a class="external text external" href="https://en.wikipedia.org/w/index.php?title=Transformer_(deep_learning_architecture)&amp;action=edit">edit</a><span class="mw-editsection-bracket">]</span></span></div><div class="excerpt">
<p>Transformers can also be used/adapted for modalities (input or output) beyond just text, usually by finding a way to "tokenize" the modality.
</p><p>Multimodal models can either be trained from scratch, or by finetuning. A 2022 study found that Transformers pretrained only on natural language can be finetuned on only 0.03% of parameters and become competitive with <a href="LSTMs" class="mw-redirect" title="LSTMs">LSTMs</a> on a variety of logical and visual tasks, demonstrating <a href="Transfer_learning" title="Transfer learning">transfer learning</a>.<sup id="cite_ref-6" class="reference"><a href="#cite_note-6"><span class="cite-bracket">[</span>6<span class="cite-bracket">]</span></a></sup> The LLaVA was a vision-language model composed of a language model (Vicuna-13B)<sup id="cite_ref-7" class="reference"><a href="#cite_note-7"><span class="cite-bracket">[</span>7<span class="cite-bracket">]</span></a></sup> and a vision model (<a href="Vision_transformer" title="Vision transformer">ViT</a>-L/14), connected by a linear layer. Only the linear layer is finetuned.<sup id="cite_ref-8" class="reference"><a href="#cite_note-8"><span class="cite-bracket">[</span>8<span class="cite-bracket">]</span></a></sup>
</p><p><a href="Vision_transformer" title="Vision transformer">Vision transformers</a><sup id="cite_ref-Transformer_(deep_learning_architecture)_auto2_9-0" class="reference"><a href="#cite_note-Transformer_(deep_learning_architecture)_auto2-9"><span class="cite-bracket">[</span>9<span class="cite-bracket">]</span></a></sup> adapt the transformer to computer vision by breaking down input images as a series of patches, turning them into vectors, and treating them like tokens in a standard transformer.
</p><p>Conformer<sup id="cite_ref-Transformer_(deep_learning_architecture)_Gulati2020_10-0" class="reference"><a href="#cite_note-Transformer_(deep_learning_architecture)_Gulati2020-10"><span class="cite-bracket">[</span>10<span class="cite-bracket">]</span></a></sup> and later <a href="Whisper_(speech_recognition_system)" title="Whisper (speech recognition system)">Whisper</a><sup id="cite_ref-Transformer_(deep_learning_architecture)_Radford_Kim_Xu_Brockman_p._11-0" class="reference"><a href="#cite_note-Transformer_(deep_learning_architecture)_Radford_Kim_Xu_Brockman_p.-11"><span class="cite-bracket">[</span>11<span class="cite-bracket">]</span></a></sup> follow the same pattern for <a href="Speech_recognition" title="Speech recognition">speech recognition</a>, first turning the speech signal into a <a href="Spectrogram" title="Spectrogram">spectrogram</a>, which is then treated like an image, i.e. broken down into a series of patches, turned into vectors and treated like tokens in a standard transformer.
</p><p><a href="Perceiver" title="Perceiver">Perceivers</a><sup id="cite_ref-Transformer_(deep_learning_architecture)_perceiver2021_12-0" class="reference"><a href="#cite_note-Transformer_(deep_learning_architecture)_perceiver2021-12"><span class="cite-bracket">[</span>12<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-Transformer_(deep_learning_architecture)_jaegle2021b_13-0" class="reference"><a href="#cite_note-Transformer_(deep_learning_architecture)_jaegle2021b-13"><span class="cite-bracket">[</span>13<span class="cite-bracket">]</span></a></sup> are a variant of Transformers designed for multimodality.
</p>
For image generation, notable architectures are <a href="DALL-E" title="DALL-E">DALL-E 1</a> (2021), Parti (2022),<sup id="cite_ref-14" class="reference"><a href="#cite_note-14"><span class="cite-bracket">[</span>14<span class="cite-bracket">]</span></a></sup> Phenaki (2023),<sup id="cite_ref-Transformer_(deep_learning_architecture)_:13_15-0" class="reference"><a href="#cite_note-Transformer_(deep_learning_architecture)_:13-15"><span class="cite-bracket">[</span>15<span class="cite-bracket">]</span></a></sup> and Muse (2023).<sup id="cite_ref-Transformer_(deep_learning_architecture)_:12_16-0" class="reference"><a href="#cite_note-Transformer_(deep_learning_architecture)_:12-16"><span class="cite-bracket">[</span>16<span class="cite-bracket">]</span></a></sup> Unlike later models, DALL-E is not a diffusion model. Instead, it uses a decoder-only Transformer that autoregressively generates a text, followed by the token representation of an image, which is then converted by a <a href="Variational_autoencoder" title="Variational autoencoder">variational autoencoder</a> to an image.<sup id="cite_ref-17" class="reference"><a href="#cite_note-17"><span class="cite-bracket">[</span>17<span class="cite-bracket">]</span></a></sup> Parti is an encoder-decoder Transformer, where the encoder processes a text prompt, and the decoder generates a token representation of an image.<sup id="cite_ref-18" class="reference"><a href="#cite_note-18"><span class="cite-bracket">[</span>18<span class="cite-bracket">]</span></a></sup> Muse is an encoder-only Transformer that is trained to predict masked image tokens from unmasked image tokens. During generation, all input tokens are masked, and the highest-confidence predictions are included for the next iteration, until all tokens are predicted.<sup id="cite_ref-Transformer_(deep_learning_architecture)_:12_16-1" class="reference"><a href="#cite_note-Transformer_(deep_learning_architecture)_:12-16"><span class="cite-bracket">[</span>16<span class="cite-bracket">]</span></a></sup> Phenaki is a text-to-video model. It is a bidirectional masked transformer conditioned on pre-computed text tokens. The generated tokens are then decoded to a video.<sup id="cite_ref-Transformer_(deep_learning_architecture)_:13_15-1" class="reference"><a href="#cite_note-Transformer_(deep_learning_architecture)_:13-15"><span class="cite-bracket">[</span>15<span class="cite-bracket">]</span></a></sup></div></div>
<div class="mw-heading mw-heading3"><h3 id="Multimodal_large_language_models">Multimodal large language models</h3></div>
<div class="excerpt-block"><div role="note" class="hatnote navigation-not-searchable dablink excerpt-hat selfref">This section is an excerpt from <a href="Large_language_model#Multimodality" title="Large language model">Large language model § Multimodality</a>.<span class="mw-editsection-like "><span class="mw-editsection-bracket">[</span><a class="external text external" href="https://en.wikipedia.org/w/index.php?title=Large_language_model&amp;action=edit">edit</a><span class="mw-editsection-bracket">]</span></span></div><div class="excerpt">
<p>Multimodality means having multiple modalities, where a "<a href="Modality_(human%E2%80%93computer_interaction)" title="Modality (human–computer interaction)">modality</a>" refers to a type of input or output, such as video, image, audio, text, <a href="Proprioception" title="Proprioception">proprioception</a>, etc.<sup id="cite_ref-19" class="reference"><a href="#cite_note-19"><span class="cite-bracket">[</span>19<span class="cite-bracket">]</span></a></sup> For example, <a href="Pathways_Language_Model" class="mw-redirect" title="Pathways Language Model">Google PaLM</a> model was fine-tuned into a multimodal model and applied to <a href="Robot_control" title="Robot control">robotic control</a>.<sup id="cite_ref-20" class="reference"><a href="#cite_note-20"><span class="cite-bracket">[</span>20<span class="cite-bracket">]</span></a></sup> <a href="LLaMA" class="mw-redirect" title="LLaMA">LLaMA</a> models have also been turned multimodal using the tokenization method, to allow image inputs,<sup id="cite_ref-21" class="reference"><a href="#cite_note-21"><span class="cite-bracket">[</span>21<span class="cite-bracket">]</span></a></sup> and video inputs.<sup id="cite_ref-22" class="reference"><a href="#cite_note-22"><span class="cite-bracket">[</span>22<span class="cite-bracket">]</span></a></sup> <a href="GPT-4o" title="GPT-4o">GPT-4o</a> can process and generate text, audio and images.<sup id="cite_ref-23" class="reference"><a href="#cite_note-23"><span class="cite-bracket">[</span>23<span class="cite-bracket">]</span></a></sup> Such models are sometimes called large multimodal models (LMMs).<sup id="cite_ref-24" class="reference"><a href="#cite_note-24"><span class="cite-bracket">[</span>24<span class="cite-bracket">]</span></a></sup>
</p>
A common method to create multimodal models out of an LLM is to "tokenize" the output of a trained encoder. Concretely, one can construct an LLM that can understand images as follows: take a trained LLM, and take a trained image encoder <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle E}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>E</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle E}</annotation>
</semantics>
</math></span><img src="./4232c9de2ee3eec0a9c0a19b15ab92daa6223f9b.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.338ex; width:1.776ex; height:2.176ex;" alt="{\displaystyle E}" loading="lazy"></span>. Make a small multilayered perceptron <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle f}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>f</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle f}</annotation>
</semantics>
</math></span><img src="./132e57acb643253e7810ee9702d9581f159a1c61.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:1.279ex; height:2.509ex;" alt="{\displaystyle f}" loading="lazy"></span>, so that for any image <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle y}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>y</mi>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle y}</annotation>
</semantics>
</math></span><img src="./b8a6208ec717213d4317e666f1ae872e00620a0d.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.671ex; width:1.155ex; height:2.009ex;" alt="{\displaystyle y}" loading="lazy"></span>, the post-processed vector <span class="mwe-math-element mwe-math-element-inline"><span class="mwe-math-mathml-inline mwe-math-mathml-a11y" style="display: none;"><math xmlns="http://www.w3.org/1998/Math/MathML" alttext="{\displaystyle f(E(y))}">
<semantics>
<mrow class="MJX-TeXAtom-ORD">
<mstyle displaystyle="true" scriptlevel="0">
<mi>f</mi>
<mo stretchy="false">(</mo>
<mi>E</mi>
<mo stretchy="false">(</mo>
<mi>y</mi>
<mo stretchy="false">)</mo>
<mo stretchy="false">)</mo>
</mstyle>
</mrow>
<annotation encoding="application/x-tex">{\displaystyle f(E(y))}</annotation>
</semantics>
</math></span><img src="./8d41d0ec0611a795f65ea14a43b8016462703a8e.svg" class="mwe-math-fallback-image-inline mw-invert skin-invert" aria-hidden="true" style="vertical-align: -0.838ex; width:7.828ex; height:2.843ex;" alt="{\displaystyle f(E(y))}" loading="lazy"></span> has the same dimensions as an encoded token. That is an "image token". Then, one can interleave text tokens and image tokens. The compound model is then fine-tuned on an image-text dataset. This basic construction can be applied with more sophistication to improve the model. The image encoder may be frozen to improve stability.<sup id="cite_ref-25" class="reference"><a href="#cite_note-25"><span class="cite-bracket">[</span>25<span class="cite-bracket">]</span></a></sup> The model Flamingo demonstrated in 2022 the effectiveness of the tokenization method, fine-tuning a pair of pretrained language model and image encoder to perform better on visual question answering than models trained from scratch.<sup id="cite_ref-26" class="reference"><a href="#cite_note-26"><span class="cite-bracket">[</span>26<span class="cite-bracket">]</span></a></sup></div></div>
<div class="mw-heading mw-heading2"><h2 id="Multimodal_deep_Boltzmann_machines">Multimodal deep Boltzmann machines</h2></div>
<p>A <a href="Boltzmann_machine" title="Boltzmann machine">Boltzmann machine</a> is a type of <a href="Stochastic_neural_network" class="mw-redirect" title="Stochastic neural network">stochastic neural network</a> invented by <a href="Geoffrey_Hinton" title="Geoffrey Hinton">Geoffrey Hinton</a> and <a href="Terry_Sejnowski" title="Terry Sejnowski">Terry Sejnowski</a> in 1985. Boltzmann machines can be seen as the <a href="Stochastic_process" title="Stochastic process">stochastic</a>, <a href="Generative_model" title="Generative model">generative</a> counterpart of <a href="Hopfield_net" class="mw-redirect" title="Hopfield net">Hopfield nets</a>. They are named after the <a href="Boltzmann_distribution" title="Boltzmann distribution">Boltzmann distribution</a> in statistical mechanics. The units in Boltzmann machines are divided into two groups: visible units and hidden units. Each unit is like a neuron with a binary output that represents whether it is activated or not.<sup id="cite_ref-27" class="reference"><a href="#cite_note-27"><span class="cite-bracket">[</span>27<span class="cite-bracket">]</span></a></sup> General Boltzmann machines allow connection between any units. However, learning is impractical using general Boltzmann Machines because the computational time is exponential to the size of the machine. A more efficient architecture is called <a href="Restricted_Boltzmann_machine" title="Restricted Boltzmann machine">restricted Boltzmann machine</a> where connection is only allowed between hidden unit and visible unit, which is described in the next section.
</p><p>Multimodal deep Boltzmann machines can process and learn from different types of information, such as images and text, simultaneously. This can notably be done by having a separate deep Boltzmann machine for each modality, for example one for images and one for text, joined at an additional top hidden layer.<sup id="cite_ref-28" class="reference"><a href="#cite_note-28"><span class="cite-bracket">[</span>28<span class="cite-bracket">]</span></a></sup>
</p>
<div class="mw-heading mw-heading2"><h2 id="Applications">Applications</h2></div>
<p>Multimodal machine learning has numerous applications across various domains:
</p>
<ul><li><b>Cross-modal retrieval</b>: cross-modal retrieval allows users to search for data across different modalities (e.g., retrieving images based on text descriptions), improving multimedia search engines and content recommendation systems. Models like <a href="Contrastive_Language-Image_Pre-training" title="Contrastive Language-Image Pre-training">CLIP</a> facilitate efficient, accurate retrieval by embedding data in a shared space, demonstrating strong performance even in zero-shot settings.<sup id="cite_ref-29" class="reference"><a href="#cite_note-29"><span class="cite-bracket">[</span>29<span class="cite-bracket">]</span></a></sup></li>
<li><b>Classification and missing data retrieval</b>: multimodal Deep Boltzmann Machines outperform traditional models like <a href="Support_vector_machine" title="Support vector machine">support vector machines</a> and <a href="Latent_Dirichlet_allocation" title="Latent Dirichlet allocation">latent Dirichlet allocation</a> in classification tasks and can predict missing data in multimodal datasets, such as images and text.</li>
<li><b>Healthcare diagnostics</b>: multimodal models integrate medical imaging, genomic data, and patient records to improve diagnostic accuracy and early disease detection, especially in cancer screening.<sup id="cite_ref-30" class="reference"><a href="#cite_note-30"><span class="cite-bracket">[</span>30<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-31" class="reference"><a href="#cite_note-31"><span class="cite-bracket">[</span>31<span class="cite-bracket">]</span></a></sup><sup id="cite_ref-32" class="reference"><a href="#cite_note-32"><span class="cite-bracket">[</span>32<span class="cite-bracket">]</span></a></sup></li>
<li><b>Content generation</b>: models like <a href="DALL%C2%B7E" class="mw-redirect" title="DALL·E">DALL·E</a> generate images from textual descriptions, benefiting creative industries, while cross-modal retrieval enables dynamic multimedia searches.<sup id="cite_ref-33" class="reference"><a href="#cite_note-33"><span class="cite-bracket">[</span>33<span class="cite-bracket">]</span></a></sup></li>
<li><b>Robotics and human-computer interaction</b>: multimodal learning improves interaction in robotics and AI by integrating sensory inputs like speech, vision, and touch, aiding autonomous systems and <a href="Human%E2%80%93computer_interaction" title="Human–computer interaction">human-computer interaction</a>.</li>
<li><b>Emotion recognition</b>: combining visual, audio, and text data, multimodal systems enhance <a href="Sentiment_analysis" title="Sentiment analysis">sentiment analysis</a> and <a href="Emotion_recognition" title="Emotion recognition">emotion recognition</a>, applied in customer service, social media, and marketing.</li></ul>
<div class="mw-heading mw-heading2"><h2 id="See_also">See also</h2></div>
<ul><li><a href="Hopfield_network" title="Hopfield network">Hopfield network</a></li>
<li><a href="Markov_random_field" title="Markov random field">Markov random field</a></li>
<li><a href="Markov_chain_Monte_Carlo" title="Markov chain Monte Carlo">Markov chain Monte Carlo</a></li></ul>
<div class="mw-heading mw-heading2"><h2 id="References">References</h2></div>
<style data-mw-deduplicate="TemplateStyles:r1239543626">
/* start https://en.wikipedia.org/ */


.mw-parser-output .reflist{margin-bottom:0.5em;list-style-type:decimal}@media screen{.mw-parser-output .reflist{font-size:90%}}.mw-parser-output .reflist .references{font-size:100%;margin-bottom:0;list-style-type:inherit}.mw-parser-output .reflist-columns-2{column-width:30em}.mw-parser-output .reflist-columns-3{column-width:25em}.mw-parser-output .reflist-columns{margin-top:0.3em}.mw-parser-output .reflist-columns ol{margin-top:0}.mw-parser-output .reflist-columns li{page-break-inside:avoid;break-inside:avoid-column}.mw-parser-output .reflist-upper-alpha{list-style-type:upper-alpha}.mw-parser-output .reflist-upper-roman{list-style-type:upper-roman}.mw-parser-output .reflist-lower-alpha{list-style-type:lower-alpha}.mw-parser-output .reflist-lower-greek{list-style-type:lower-greek}.mw-parser-output .reflist-lower-roman{list-style-type:lower-roman}


/* end https://en.wikipedia.org/ */
</style><div class="reflist">
<div class="mw-references-wrap mw-references-columns"><ol class="references">
<li id="cite_note-1"><span class="mw-cite-backlink"><b><a href="#cite_ref-1">^</a></b></span> <span class="reference-text"><style data-mw-deduplicate="TemplateStyles:r1238218222">
/* start https://en.wikipedia.org/ */


.mw-parser-output cite.citation{font-style:inherit;word-wrap:break-word}.mw-parser-output .citation q{quotes:"\"""\"""'""'"}.mw-parser-output .citation:target{background-color:rgba(0,127,255,0.133)}.mw-parser-output .id-lock-free.id-lock-free a{background:url("./mw/Lock-green.svg")right 0.1em center/9px no-repeat}.mw-parser-output .id-lock-limited.id-lock-limited a,.mw-parser-output .id-lock-registration.id-lock-registration a{background:url("./mw/Lock-gray-alt-2.svg")right 0.1em center/9px no-repeat}.mw-parser-output .id-lock-subscription.id-lock-subscription a{background:url("./mw/Lock-red-alt-2.svg")right 0.1em center/9px no-repeat}.mw-parser-output .cs1-ws-icon a{background:url("./mw/Wikisource-logo.svg")right 0.1em center/12px no-repeat}body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-free a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-limited a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-registration a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .id-lock-subscription a,body:not(.skin-timeless):not(.skin-minerva) .mw-parser-output .cs1-ws-icon a{background-size:contain;padding:0 1em 0 0}.mw-parser-output .cs1-code{color:inherit;background:inherit;border:none;padding:inherit}.mw-parser-output .cs1-hidden-error{display:none;color:var(--color-error,#d33)}.mw-parser-output .cs1-visible-error{color:var(--color-error,#d33)}.mw-parser-output .cs1-maint{display:none;color:#085;margin-left:0.3em}.mw-parser-output .cs1-kern-left{padding-left:0.2em}.mw-parser-output .cs1-kern-right{padding-right:0.2em}.mw-parser-output .citation .mw-selflink{font-weight:inherit}@media screen{.mw-parser-output .cs1-format{font-size:95%}html.skin-theme-clientpref-night .mw-parser-output .cs1-maint{color:#18911f}}@media screen and (prefers-color-scheme:dark){html.skin-theme-clientpref-os .mw-parser-output .cs1-maint{color:#18911f}}


/* end https://en.wikipedia.org/ */
</style><cite id="CITEREFHendriksenBleekerVakulenkovan_Noord2021" class="citation arxiv cs1">Hendriksen, Mariya; Bleeker, Maurits; Vakulenko, Svitlana; van Noord, Nanne; Kuiper, Ernst; de Rijke, Maarten (2021). "Extending CLIP for Category-to-image Retrieval in E-commerce". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2112.11294">2112.11294</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CV">cs.CV</a>].</cite></span>
</li>
<li id="cite_note-stable-diffusion-github-2"><span class="mw-cite-backlink"><b><a href="#cite_ref-stable-diffusion-github_2-0">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://github.com/CompVis/stable-diffusion">"Stable Diffusion Repository on GitHub"</a>. CompVis - Machine Vision and Learning Research Group, LMU Munich. 17 September 2022. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230118183342/https://github.com/CompVis/stable-diffusion">Archived</a> from the original on January 18, 2023<span class="reference-accessdate">. Retrieved <span class="nowrap">17 September</span> 2022</span>.</cite></span>
</li>
<li id="cite_note-3"><span class="mw-cite-backlink"><b><a href="#cite_ref-3">^</a></b></span> <span class="reference-text"><cite class="citation cs2"><a rel="nofollow" class="external text" href="https://github.com/LAION-AI/aesthetic-predictor"><i>LAION-AI/aesthetic-predictor</i></a>, LAION AI, 2024-09-06<span class="reference-accessdate">, retrieved <span class="nowrap">2024-09-08</span></span></cite></span>
</li>
<li id="cite_note-4"><span class="mw-cite-backlink"><b><a href="#cite_ref-4">^</a></b></span> <span class="reference-text"><cite id="CITEREFMokadyHertzBermano2021" class="citation arxiv cs1">Mokady, Ron; Hertz, Amir; Bermano, Amit H. (2021). "ClipCap: CLIP Prefix for Image Captioning". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2111.09734">2111.09734</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CV">cs.CV</a>].</cite></span>
</li>
<li id="cite_note-5"><span class="mw-cite-backlink"><b><a href="#cite_ref-5">^</a></b></span> <span class="reference-text"><cite id="CITEREFZia2024" class="citation web cs1">Zia, Tehseen (January 8, 2024). <a rel="nofollow" class="external text" href="https://www.unite.ai/unveiling-of-large-multimodal-models-shaping-the-landscape-of-language-models-in-2024/">"Unveiling of Large Multimodal Models: Shaping the Landscape of Language Models in 2024"</a>. <i>Unite.ai</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2024-06-01</span></span>.</cite></span>
</li>
<li id="cite_note-6"><span class="mw-cite-backlink"><b><a href="#cite_ref-6">^</a></b></span> <span class="reference-text"><cite id="CITEREFLuGroverAbbeelMordatch2022" class="citation journal cs1">Lu, Kevin; Grover, Aditya; Abbeel, Pieter; Mordatch, Igor (2022-06-28). <a rel="nofollow" class="external text" href="https://ojs.aaai.org/index.php/AAAI/article/view/20729">"Frozen Pretrained Transformers as Universal Computation Engines"</a>. <i>Proceedings of the AAAI Conference on Artificial Intelligence</i>. <b>36</b> (7): <span class="nowrap">7628–</span>7636. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1609%2Faaai.v36i7.20729">10.1609/aaai.v36i7.20729</a></span>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a>&nbsp;<a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/2374-3468">2374-3468</a>.</cite></span>
</li>
<li id="cite_note-7"><span class="mw-cite-backlink"><b><a href="#cite_ref-7">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://lmsys.org/blog/2023-03-30-vicuna">"Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality | LMSYS Org"</a>. <i>lmsys.org</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2024-08-11</span></span>.</cite></span>
</li>
<li id="cite_note-8"><span class="mw-cite-backlink"><b><a href="#cite_ref-8">^</a></b></span> <span class="reference-text"><cite id="CITEREFLiuLiWuLee2023" class="citation journal cs1">Liu, Haotian; Li, Chunyuan; Wu, Qingyang; Lee, Yong Jae (2023-12-15). <a rel="nofollow" class="external text" href="https://proceedings.neurips.cc/paper_files/paper/2023/hash/6dcf277ea32ce3288914faf369fe6de0-Abstract-Conference.html">"Visual Instruction Tuning"</a>. <i>Advances in Neural Information Processing Systems</i>. <b>36</b>: <span class="nowrap">34892–</span>34916.</cite></span>
</li>
<li id="cite_note-Transformer_(deep_learning_architecture)_auto2-9"><span class="mw-cite-backlink"><b><a href="#cite_ref-Transformer_(deep_learning_architecture)_auto2_9-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFDosovitskiyBeyerKolesnikovWeissenborn2021" class="citation arxiv cs1">Dosovitskiy, Alexey; Beyer, Lucas; Kolesnikov, Alexander; Weissenborn, Dirk; Zhai, Xiaohua; Unterthiner, Thomas; Dehghani, Mostafa; Minderer, Matthias; Heigold, Georg; Gelly, Sylvain; Uszkoreit, Jakob (2021-06-03). "An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2010.11929">2010.11929</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CV">cs.CV</a>].</cite></span>
</li>
<li id="cite_note-Transformer_(deep_learning_architecture)_Gulati2020-10"><span class="mw-cite-backlink"><b><a href="#cite_ref-Transformer_(deep_learning_architecture)_Gulati2020_10-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFGulatiQinChiuParmar2020" class="citation arxiv cs1">Gulati, Anmol; Qin, James; Chiu, Chung-Cheng; Parmar, Niki; Zhang, Yu; Yu, Jiahui; Han, Wei; Wang, Shibo; Zhang, Zhengdong; Wu, Yonghui; Pang, Ruoming (2020). "Conformer: Convolution-augmented Transformer for Speech Recognition". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2005.08100">2005.08100</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/eess.AS">eess.AS</a>].</cite></span>
</li>
<li id="cite_note-Transformer_(deep_learning_architecture)_Radford_Kim_Xu_Brockman_p.-11"><span class="mw-cite-backlink"><b><a href="#cite_ref-Transformer_(deep_learning_architecture)_Radford_Kim_Xu_Brockman_p._11-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFRadfordKimXuBrockman2022" class="citation arxiv cs1">Radford, Alec; Kim, Jong Wook; Xu, Tao; Brockman, Greg; McLeavey, Christine; Sutskever, Ilya (2022). "Robust Speech Recognition via Large-Scale Weak Supervision". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2212.04356">2212.04356</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/eess.AS">eess.AS</a>].</cite></span>
</li>
<li id="cite_note-Transformer_(deep_learning_architecture)_perceiver2021-12"><span class="mw-cite-backlink"><b><a href="#cite_ref-Transformer_(deep_learning_architecture)_perceiver2021_12-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFJaegleGimenoBrockZisserman2021" class="citation arxiv cs1">Jaegle, Andrew; Gimeno, Felix; Brock, Andrew; Zisserman, Andrew; Vinyals, Oriol; Carreira, Joao (2021-06-22). "Perceiver: General Perception with Iterative Attention". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2103.03206">2103.03206</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CV">cs.CV</a>].</cite></span>
</li>
<li id="cite_note-Transformer_(deep_learning_architecture)_jaegle2021b-13"><span class="mw-cite-backlink"><b><a href="#cite_ref-Transformer_(deep_learning_architecture)_jaegle2021b_13-0">^</a></b></span> <span class="reference-text"><cite id="CITEREFJaegleBorgeaudAlayracDoersch2021" class="citation arxiv cs1">Jaegle, Andrew; Borgeaud, Sebastian; Alayrac, Jean-Baptiste; Doersch, Carl; Ionescu, Catalin; Ding, David; Koppula, Skanda; Zoran, Daniel; Brock, Andrew; Shelhamer, Evan; Hénaff, Olivier (2021-08-02). "Perceiver IO: A General Architecture for Structured Inputs &amp; Outputs". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2107.14795">2107.14795</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.LG">cs.LG</a>].</cite></span>
</li>
<li id="cite_note-14"><span class="mw-cite-backlink"><b><a href="#cite_ref-14">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="https://sites.research.google/parti/">"Parti: Pathways Autoregressive Text-to-Image Model"</a>. <i>sites.research.google</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2024-08-09</span></span>.</cite></span>
</li>
<li id="cite_note-Transformer_(deep_learning_architecture)_:13-15"><span class="mw-cite-backlink">^ <a href="#cite_ref-Transformer_(deep_learning_architecture)_:13_15-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-Transformer_(deep_learning_architecture)_:13_15-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFVillegasBabaeizadehKindermansMoraldo2022" class="citation journal cs1">Villegas, Ruben; Babaeizadeh, Mohammad; Kindermans, Pieter-Jan; Moraldo, Hernan; Zhang, Han; Saffar, Mohammad Taghi; Castro, Santiago; Kunze, Julius; Erhan, Dumitru (2022-09-29). <a rel="nofollow" class="external text" href="https://openreview.net/forum?id=vOEXS39nOF">"Phenaki: Variable Length Video Generation from Open Domain Textual Descriptions"</a>.</cite> <span class="cs1-visible-error citation-comment"><code class="cs1-code">{{cite journal}}</code>: </span><span class="cs1-visible-error citation-comment">Cite journal requires <code class="cs1-code">|journal=</code> (help)</span></span>
</li>
<li id="cite_note-Transformer_(deep_learning_architecture)_:12-16"><span class="mw-cite-backlink">^ <a href="#cite_ref-Transformer_(deep_learning_architecture)_:12_16-0"><sup><i><b>a</b></i></sup></a> <a href="#cite_ref-Transformer_(deep_learning_architecture)_:12_16-1"><sup><i><b>b</b></i></sup></a></span> <span class="reference-text"><cite id="CITEREFChangZhangBarberMaschinot2023" class="citation arxiv cs1">Chang, Huiwen; Zhang, Han; Barber, Jarred; Maschinot, A. J.; Lezama, Jose; Jiang, Lu; Yang, Ming-Hsuan; Murphy, Kevin; Freeman, William T. (2023-01-02). "Muse: Text-To-Image Generation via Masked Generative Transformers". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2301.00704">2301.00704</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CV">cs.CV</a>].</cite></span>
</li>
<li id="cite_note-17"><span class="mw-cite-backlink"><b><a href="#cite_ref-17">^</a></b></span> <span class="reference-text"><cite id="CITEREFRameshPavlovGohGray2021" class="citation cs2">Ramesh, Aditya; Pavlov, Mikhail; Goh, Gabriel; Gray, Scott; Voss, Chelsea; Radford, Alec; Chen, Mark; Sutskever, Ilya (2021-02-26), <i>Zero-Shot Text-to-Image Generation</i>, <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2102.12092">2102.12092</a></span></cite></span>
</li>
<li id="cite_note-18"><span class="mw-cite-backlink"><b><a href="#cite_ref-18">^</a></b></span> <span class="reference-text"><cite id="CITEREFYuXuKohLuong2022" class="citation cs2">Yu, Jiahui; Xu, Yuanzhong; Koh, Jing Yu; Luong, Thang; Baid, Gunjan; Wang, Zirui; Vasudevan, Vijay; Ku, Alexander; Yang, Yinfei (2022-06-21), <i>Scaling Autoregressive Models for Content-Rich Text-to-Image Generation</i>, <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2206.10789">2206.10789</a></span></cite></span>
</li>
<li id="cite_note-19"><span class="mw-cite-backlink"><b><a href="#cite_ref-19">^</a></b></span> <span class="reference-text"><cite id="CITEREFKirosSalakhutdinovZemel2014" class="citation journal cs1">Kiros, Ryan; Salakhutdinov, Ruslan; Zemel, Rich (2014-06-18). <a rel="nofollow" class="external text" href="https://proceedings.mlr.press/v32/kiros14.html">"Multimodal Neural Language Models"</a>. <i>Proceedings of the 31st International Conference on Machine Learning</i>. PMLR: <span class="nowrap">595–</span>603. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230702195952/https://proceedings.mlr.press/v32/kiros14.html">Archived</a> from the original on 2023-07-02<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-07-02</span></span>.</cite></span>
</li>
<li id="cite_note-20"><span class="mw-cite-backlink"><b><a href="#cite_ref-20">^</a></b></span> <span class="reference-text"><cite id="CITEREFDriessXiaSajjadiLynch2023" class="citation arxiv cs1">Driess, Danny; Xia, Fei; Sajjadi, Mehdi S. M.; Lynch, Corey; Chowdhery, Aakanksha; Ichter, Brian; Wahid, Ayzaan; Tompson, Jonathan; Vuong, Quan; Yu, Tianhe; Huang, Wenlong; Chebotar, Yevgen; Sermanet, Pierre; Duckworth, Daniel; Levine, Sergey (2023-03-01). "PaLM-E: An Embodied Multimodal Language Model". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2303.03378">2303.03378</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.LG">cs.LG</a>].</cite></span>
</li>
<li id="cite_note-21"><span class="mw-cite-backlink"><b><a href="#cite_ref-21">^</a></b></span> <span class="reference-text"><cite id="CITEREFLiuLiWuLee2023" class="citation arxiv cs1">Liu, Haotian; Li, Chunyuan; Wu, Qingyang; Lee, Yong Jae (2023-04-01). "Visual Instruction Tuning". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2304.08485">2304.08485</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CV">cs.CV</a>].</cite></span>
</li>
<li id="cite_note-22"><span class="mw-cite-backlink"><b><a href="#cite_ref-22">^</a></b></span> <span class="reference-text"><cite id="CITEREFZhangLiBing2023" class="citation arxiv cs1">Zhang, Hang; Li, Xin; Bing, Lidong (2023-06-01). "Video-LLaMA: An Instruction-tuned Audio-Visual Language Model for Video Understanding". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2306.02858">2306.02858</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CL">cs.CL</a>].</cite></span>
</li>
<li id="cite_note-23"><span class="mw-cite-backlink"><b><a href="#cite_ref-23">^</a></b></span> <span class="reference-text"><cite class="citation news cs1"><a rel="nofollow" class="external text" href="https://www.theregister.com/2024/05/13/openai_gpt4o/">"OpenAI says natively multimodal GPT-4o eats text, visuals, sound – and emits the same"</a>. <i>The Register</i>. 2024-05-13.</cite></span>
</li>
<li id="cite_note-24"><span class="mw-cite-backlink"><b><a href="#cite_ref-24">^</a></b></span> <span class="reference-text"><cite id="CITEREFZia2024" class="citation web cs1">Zia, Dr Tehseen (2024-01-08). <a rel="nofollow" class="external text" href="https://www.unite.ai/unveiling-of-large-multimodal-models-shaping-the-landscape-of-language-models-in-2024/">"Unveiling of Large Multimodal Models: Shaping the Landscape of Language Models in 2024"</a>. <i>Unite.AI</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2025-05-30</span></span>.</cite></span>
</li>
<li id="cite_note-25"><span class="mw-cite-backlink"><b><a href="#cite_ref-25">^</a></b></span> <span class="reference-text"><cite id="CITEREFLiLiSavareseHoi2023" class="citation arxiv cs1">Li, Junnan; Li, Dongxu; Savarese, Silvio; Hoi, Steven (2023-01-01). "BLIP-2: Bootstrapping Language-Image Pre-training with Frozen Image Encoders and Large Language Models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2301.12597">2301.12597</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CV">cs.CV</a>].</cite></span>
</li>
<li id="cite_note-26"><span class="mw-cite-backlink"><b><a href="#cite_ref-26">^</a></b></span> <span class="reference-text"><cite id="CITEREFAlayracDonahueLucMiech2022" class="citation journal cs1">Alayrac, Jean-Baptiste; Donahue, Jeff; Luc, Pauline; Miech, Antoine; Barr, Iain; Hasson, Yana; Lenc, Karel; Mensch, Arthur; Millican, Katherine; Reynolds, Malcolm; Ring, Roman; Rutherford, Eliza; Cabi, Serkan; Han, Tengda; Gong, Zhitao (2022-12-06). <a rel="nofollow" class="external text" href="https://proceedings.neurips.cc/paper_files/paper/2022/hash/960a172bc7fbf0177ccccbb411a7d800-Abstract-Conference.html">"Flamingo: a Visual Language Model for Few-Shot Learning"</a>. <i>Advances in Neural Information Processing Systems</i>. <b>35</b>: <span class="nowrap">23716–</span>23736. <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2204.14198">2204.14198</a></span>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20230702195951/https://proceedings.neurips.cc/paper_files/paper/2022/hash/960a172bc7fbf0177ccccbb411a7d800-Abstract-Conference.html">Archived</a> from the original on 2023-07-02<span class="reference-accessdate">. Retrieved <span class="nowrap">2023-07-02</span></span>.</cite></span>
</li>
<li id="cite_note-27"><span class="mw-cite-backlink"><b><a href="#cite_ref-27">^</a></b></span> <span class="reference-text"><cite id="CITEREFDey2021" class="citation web cs1">Dey, Victor (2021-09-03). <a rel="nofollow" class="external text" href="https://analyticsindiamag.com/beginners-guide-to-boltzmann-machines/">"Beginners Guide to Boltzmann Machine"</a>. <i>Analytics India Magazine</i><span class="reference-accessdate">. Retrieved <span class="nowrap">2024-03-02</span></span>.</cite></span>
</li>
<li id="cite_note-28"><span class="mw-cite-backlink"><b><a href="#cite_ref-28">^</a></b></span> <span class="reference-text"><cite class="citation web cs1"><a rel="nofollow" class="external text" href="http://www.jmlr.org/papers/volume15/srivastava14b/srivastava14b.pdf">"Multimodal Learning with Deep Boltzmann Machine"</a> <span class="cs1-format">(PDF)</span>. 2014. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20150621055730/http://jmlr.org/papers/volume15/srivastava14b/srivastava14b.pdf">Archived</a> <span class="cs1-format">(PDF)</span> from the original on 2015-06-21<span class="reference-accessdate">. Retrieved <span class="nowrap">2015-06-14</span></span>.</cite></span>
</li>
<li id="cite_note-29"><span class="mw-cite-backlink"><b><a href="#cite_ref-29">^</a></b></span> <span class="reference-text"><cite id="CITEREFHendriksenVakulenkoKuiperde_Rijke2023" class="citation arxiv cs1">Hendriksen, Mariya; Vakulenko, Svitlana; Kuiper, Ernst; de Rijke, Maarten (2023). "Scene-centric vs. Object-centric Image-Text Cross-modal Retrieval: A Reproducibility Study". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/2301.05174">2301.05174</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.CV">cs.CV</a>].</cite></span>
</li>
<li id="cite_note-30"><span class="mw-cite-backlink"><b><a href="#cite_ref-30">^</a></b></span> <span class="reference-text"><cite id="CITEREFQuach" class="citation news cs1">Quach, Katyanna. <a rel="nofollow" class="external text" href="https://www.theregister.com/2022/08/09/ai_cancer_multimodal/">"Harvard boffins build multimodal AI system to predict cancer"</a>. <i>The Register</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20220920163859/https://www.theregister.com/2022/08/09/ai_cancer_multimodal/">Archived</a> from the original on 20 September 2022<span class="reference-accessdate">. Retrieved <span class="nowrap">16 September</span> 2022</span>.</cite></span>
</li>
<li id="cite_note-31"><span class="mw-cite-backlink"><b><a href="#cite_ref-31">^</a></b></span> <span class="reference-text"><cite id="CITEREFChenLuWilliamsonChen2022" class="citation journal cs1">Chen, Richard J.; Lu, Ming Y.; Williamson, Drew F. K.; Chen, Tiffany Y.; Lipkova, Jana; Noor, Zahra; Shaban, Muhammad; Shady, Maha; Williams, Mane; Joo, Bumjin; Mahmood, Faisal (8 August 2022). <a rel="nofollow" class="external text" href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC10397370">"Pan-cancer integrative histology-genomic analysis via multimodal deep learning"</a>. <i>Cancer Cell</i>. <b>40</b> (8): 865–878.e6. <a href="Doi_(identifier)" class="mw-redirect" title="Doi (identifier)">doi</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://doi.org/10.1016%2Fj.ccell.2022.07.004">10.1016/j.ccell.2022.07.004</a></span>. <a href="ISSN_(identifier)" class="mw-redirect" title="ISSN (identifier)">ISSN</a>&nbsp;<a rel="nofollow" class="external text" href="https://search.worldcat.org/issn/1535-6108">1535-6108</a>. <a href="PMC_(identifier)" class="mw-redirect" title="PMC (identifier)">PMC</a>&nbsp;<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://www.ncbi.nlm.nih.gov/pmc/articles/PMC10397370">10397370</a></span>. <a href="PMID_(identifier)" class="mw-redirect" title="PMID (identifier)">PMID</a>&nbsp;<a rel="nofollow" class="external text" href="https://pubmed.ncbi.nlm.nih.gov/35944502">35944502</a>. <a href="S2CID_(identifier)" class="mw-redirect" title="S2CID (identifier)">S2CID</a>&nbsp;<a rel="nofollow" class="external text" href="https://api.semanticscholar.org/CorpusID:251456162">251456162</a>.</cite>
<ul><li>Teaching hospital press release: <cite class="citation news cs1"><a rel="nofollow" class="external text" href="https://medicalxpress.com/news/2022-08-ai-technology-multiple-cancer-outcomes.html">"New AI technology integrates multiple data types to predict cancer outcomes"</a>. <i><a href="Brigham_and_Women's_Hospital" title="Brigham and Women's Hospital">Brigham and Women's Hospital</a> via medicalxpress.com</i>. <a rel="nofollow" class="external text" href="https://web.archive.org/web/20220920172825/https://medicalxpress.com/news/2022-08-ai-technology-multiple-cancer-outcomes.html">Archived</a> from the original on 20 September 2022<span class="reference-accessdate">. Retrieved <span class="nowrap">18 September</span> 2022</span>.</cite></li></ul>
</span></li>
<li id="cite_note-32"><span class="mw-cite-backlink"><b><a href="#cite_ref-32">^</a></b></span> <span class="reference-text"><cite id="CITEREFShiSiddharthPaigeTorr2019" class="citation arxiv cs1">Shi, Yuge; Siddharth, N.; Paige, Brooks; Torr, Philip HS (2019). "Variational Mixture-of-Experts Autoencoders for Multi-Modal Deep Generative Models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/1911.03393">1911.03393</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.LG">cs.LG</a>].</cite></span>
</li>
<li id="cite_note-33"><span class="mw-cite-backlink"><b><a href="#cite_ref-33">^</a></b></span> <span class="reference-text"><cite id="CITEREFShiSiddharthPaigeTorr2019" class="citation arxiv cs1">Shi, Yuge; Siddharth, N.; Paige, Brooks; Torr, Philip HS (2019). "Variational Mixture-of-Experts Autoencoders for Multi-Modal Deep Generative Models". <a href="ArXiv_(identifier)" class="mw-redirect" title="ArXiv (identifier)">arXiv</a>:<span class="id-lock-free" title="Freely accessible"><a rel="nofollow" class="external text" href="https://arxiv.org/abs/1911.03393">1911.03393</a></span> [<a rel="nofollow" class="external text" href="https://arxiv.org/archive/cs.LG">cs.LG</a>].</cite></span>
</li>
</ol></div></div></div><!--htdig_noindex--><div><div class="zim-footer">
This article is issued from <a class="external text" title="Last edited on 2025-06-01" href="https://en.wikipedia.org/wiki/?title=Multimodal_learning&amp;oldid=1293473618">Wikipedia</a>. The text is available under <a class="external text" href="https://creativecommons.org/licenses/by-sa/4.0/deed.en">Creative Commons Attribution-Share Alike 4.0</a> unless otherwise noted. Additional terms may apply for the media files.
</div>
</div><!--/htdig_noindex--></div>
</div>
</main>
</div>
</div>
</div>

</body></html>